import re

def sort_html_file(input_file_name, output_file_name, phrase):
    try:
        with open(input_file_name, 'r', encoding='utf-8', errors='ignore') as f:
            html = f.read()
            print("HTML loaded successfully")

        # Find all links and descriptions
        links_and_descriptions = re.findall(r'<li>.*?<\/li>', html, re.DOTALL)
        print("Links and descriptions found:", len(links_and_descriptions))

        # Extract link and description from each item
        extracted_links_and_descriptions = []
        for item in links_and_descriptions:
            match = re.search(r'href=[\'"]([^\'"]+)', item, re.DOTALL)
            if match:
                link = match.group(1)
                match = re.search(r'>[^<]+?<\/a>', item)
                if match:
                    link_text = match.group(0)[1:-4]
                    description = re.sub(r'<.*?>', '', item).strip().split('-', 1)[1].strip()
                    title_text_match = re.search(r'title=[\'"]([^\'"]+)[\'"]', item)
                    if title_text_match:
                        title_text = title_text_match.group(1)
                    else:
                        title_match = re.search(r'<a.*?>(.*?)<\/a>', item)
                        if title_text_match:
                            title_text = title_text_match.group(0)[1:-4]
                        else:
                            title_text = link_text
                    title_match = re.search(r'<li>.*?<a.*?>(.*?)<\/a>', item, re.DOTALL)
                    if title_match:
                        title = title_match.group(1)
                    else:
                        title = link_text                        
                    extracted_links_and_descriptions.append((link, link_text, title_text, description, title))

        # Add the print code here
        for link, link_text, title_text, title, description in extracted_links_and_descriptions:
         '''   print("Link:", link)
            print("Link Text:", link_text)
            print("Title Text:", title_text)
            print("Title:", title)
            print("Description:", description)
            print("------------------------") '''

        print("Extracted links and descriptions:", len(extracted_links_and_descriptions))

        # Sort links and descriptions based on the presence of the phrase
        phrase_links_and_descriptions = [x for x in extracted_links_and_descriptions if phrase.lower() in x[2].lower()]
        other_links_and_descriptions = [x for x in extracted_links_and_descriptions if phrase.lower() not in x[2].lower()]

        sorted_phrase_links_and_descriptions = sorted(phrase_links_and_descriptions, key=lambda x: x[2].lower().find(phrase.lower()))
        sorted_other_links_and_descriptions = sorted(other_links_and_descriptions, key=lambda x: x[2].lower().find(phrase.lower()))

        sorted_links_and_descriptions = []
        words = phrase.split()
        added_descriptions = set()

        for link, link_text, title_text, description, title in sorted_phrase_links_and_descriptions + sorted_other_links_and_descriptions:
            if phrase.lower() in description.lower():
                description = re.sub(r'(' + re.escape(phrase) + r')', r'<b style="background-color: #ffffb0; color: #0000FF; font-weight: bold">\1</b>', description, flags=re.IGNORECASE)
            for word in words:
                if phrase.lower() not in description.lower():
                    description = re.sub(r'(' + re.escape(word) + r')', r'<b>\1</b>', description, flags=re.IGNORECASE)
            
            description_words = description.split()
            description_hash = tuple(sorted(description_words[:9]))  # Use the first 9 words as a hash
            if description_hash not in added_descriptions:
                sorted_links_and_descriptions.append((link, link_text, title_text, description, title))
                added_descriptions.add(description_hash)

        print("Links and descriptions sorted")

        # Create the sorted HTML
        sorted_html = '<html><body><h1>Search Results</h1><ul>'
        sorted_html = '<html><body><h1>Search Results</h1>'
        sorted_html += '<meta http-equiv="content-type" content="text/html; charset=UTF-8">'
        sorted_html += '<meta charset="utf-8">'
        sorted_html += '<script>function copySentence(event) {navigator.clipboard.writeText(event.target.textContent);}</script>'
        sorted_html += '<div id="google_translate_element"></div>'
        sorted_html += '<script type="text/javascript">function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: "hi", layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, "google_translate_element");}</script>'
        sorted_html += '<script type="text/javascript" src="https://translate.google.com/translate_a/element.js?cb=googleTranslateElementInit"></script>'
        sorted_html += '<ul>'
        for link, link_text, title_text, description, title in sorted_links_and_descriptions[:10]:
            sorted_html += f'<li><a href="{link}" target="_blank" title="{title_text}" onclick="copySentence(event)">{link_text}</a> - {description}</li>'
        sorted_html += '</ul></body></html>'
        print("Sorted HTML created")

        # Save the sorted HTML to the output file
        with open(output_file_name, 'w', encoding='utf-8', errors='ignore') as f:
            f.write(sorted_html)
            print("Sorted HTML saved to output file as",output_file_name)
            print(title)
                
    except Exception as e:
        print("An error occurred:", str(e))

# Usage
input_file_name ='search_multiple_words.html'
output_file_name ='adhyatm-kya-hai.html'
phrase = 'अध्यात्म क्या है'
# sort_html_file(input_file_name, output_file_name, phrase)
